agora inbox for [email protected]help / color / mirror / Atom feed
[PATCH 2/2] Publish list of tables being repacked in shared memory 278+ messages / 2 participants [nested] [flat]
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 278+ messages in thread
* [PATCH v2 4/4] Let pgbench support binary format. @ 2026-06-02 10:08 Andy Fan <[email protected]> 0 siblings, 0 replies; 278+ messages in thread From: Andy Fan @ 2026-06-02 10:08 UTC (permalink / raw) --- src/bin/pgbench/pgbench.c | 29 +++++++++++++++++++++++++++-- 1 file changed, 27 insertions(+), 2 deletions(-) diff --git a/src/bin/pgbench/pgbench.c b/src/bin/pgbench/pgbench.c index 0b2bb9340b5..533a1732158 100644 --- a/src/bin/pgbench/pgbench.c +++ b/src/bin/pgbench/pgbench.c @@ -722,6 +722,16 @@ typedef enum QueryMode static QueryMode querymode = QUERY_SIMPLE; static const char *const QUERYMODE[] = {"simple", "extended", "prepared"}; +typedef enum ResultFormat +{ + RESULT_FORMAT_TEXT, + RESULT_FORMAT_BINARY, + NUM_RESULT_FORMAT +} ResultFormat; + +static ResultFormat result_format = RESULT_FORMAT_TEXT; +static const char *const RESULT_FORMAT[] = {"text", "binary"}; + /* * struct Command represents one command in a script. * @@ -965,6 +975,8 @@ usage(void) " --max-tries=NUM max number of tries to run transaction (default: 1)\n" " --progress-timestamp use Unix epoch timestamps for progress\n" " --random-seed=SEED set random seed (\"time\", \"rand\", integer)\n" + " --result-format=text|binary\n" + " result format for extended/prepared protocol (default: text)\n" " --sampling-rate=NUM fraction of transactions to log (e.g., 0.01 for 1%%)\n" " --show-script=NAME show builtin script code, then exit\n" " --verbose-errors print messages of all errors\n" @@ -3173,7 +3185,7 @@ sendCommand(CState *st, Command *command) pg_log_debug("client %d sending %s", st->id, sql); r = PQsendQueryParams(st->con, sql, command->argc - 1, - NULL, params, NULL, NULL, 0); + NULL, params, NULL, NULL, result_format); } else if (querymode == QUERY_PREPARED) { @@ -3184,7 +3196,7 @@ sendCommand(CState *st, Command *command) pg_log_debug("client %d sending %s", st->id, command->prepname); r = PQsendQueryPrepared(st->con, command->prepname, command->argc - 1, - params, NULL, NULL, 0); + params, NULL, NULL, result_format); } else /* unknown sql mode */ r = 0; @@ -6470,6 +6482,7 @@ printResults(StatsData *total, printf("partition method: %s\npartitions: %d\n", PARTITION_METHOD[partition_method], partitions); printf("query mode: %s\n", QUERYMODE[querymode]); + printf("result format: %s\n", RESULT_FORMAT[result_format]); printf("number of clients: %d\n", nclients); printf("number of threads: %d\n", nthreads); @@ -6778,6 +6791,7 @@ main(int argc, char **argv) {"exit-on-abort", no_argument, NULL, 16}, {"debug", no_argument, NULL, 17}, {"continue-on-error", no_argument, NULL, 18}, + {"result-format", required_argument, NULL, 19}, {NULL, 0, NULL, 0} }; @@ -7138,6 +7152,14 @@ main(int argc, char **argv) benchmarking_option_set = true; continue_on_error = true; break; + case 19: /* result-format */ + benchmarking_option_set = true; + for (result_format = 0; result_format < NUM_RESULT_FORMAT; result_format++) + if (strcmp(optarg, RESULT_FORMAT[result_format]) == 0) + break; + if (result_format >= NUM_RESULT_FORMAT) + pg_fatal("invalid result format: \"%s\"", optarg); + break; default: /* getopt_long already emitted a complaint */ pg_log_error_hint("Try \"%s --help\" for more information.", progname); @@ -7169,6 +7191,9 @@ main(int argc, char **argv) if (total_weight == 0 && !is_init_mode) pg_fatal("total script weight must not be zero"); + if (result_format == RESULT_FORMAT_BINARY && querymode == QUERY_SIMPLE) + pg_fatal("binary result format requires extended or prepared query mode"); + /* show per script stats if several scripts are used */ if (num_scripts > 1) per_script_stats = true; -- 2.43.0 --=-=-= Content-Type: application/gzip Content-Disposition: attachment; filename=test.tar.gz Content-Transfer-Encoding: base64 H4sIAAAAAAAAA+1bbXPaRhD2V/QrtgoukFhGEq9DxmmxQxI6NnhA6aRxUkYGGSsBSZWONLFDf3v3 XiQEtuMmHWg63DM2ut3bW93t7bMS5/HIJs7g3PGGl/vRH5OdtUBHVKtlejVqFT195agYO0bF1Gvl cqmCbd0wS2ZtB/T1TGcZs4jYIcCO7Y0+fcnuvv7/Kfqt49aRBSOaBkN/As963RPwA+JO3StnNHD9 AfkUONFj5b+eqMRacDHxbVJebwW4n/+lVf5X9Yrk/yYg+C/SQFaAbQPb+Pr3x3/DkPzfBNL8r0v+ bx1cj5jf4fu/Wa1K/m8Cgv8sDST7tw+48Wt++/82/td0yf9NYMF/+fa/jcCNX/Pb/zfyvyz5vwks +C/f/rcR3mzqhO5wrSXgfv6XV/hf0ivy/H8jEPyP00CWgC1D5JBZsL5HP8M9/Ddq+LIf89/AB79u GGVTnv9vBG8wAaDbGbR6vW5v0Le6p+B7ivK0hw2reXjcgvYzaL1q963+rTVBOeq1mlZL2N60UPJK JjlcoI09JpdjuczleizXUU79MYI3Y119oaN26aIl2qglzkfCVLSBcvKnTdqg/e6Uy7QhZMyBaZAo mZTuIVfLfeRKKeDS251+q2dBu2N1b1s5L61KJp8fwy5UK5VSpQAalMxatV5oNGgwoNlPjl7whmOm LQttWWjzXF2Hh2AwwhSEzM3qKTMeLiii4T41EzIaLmKaNq2vmNYT07RXEVu0LTHTWEbb1BZQ4xyN uZaDz59hjIZUolbxllCTpzRbcqZu6ppu4E8OHgELUKlaESvDT7pXdGS8eXSk1T7BkbreYD9sHBtY r5YxJhgd3IhW79fmMeQMiJyh741yBToL3DY2C7Hxsa++1Tw5TU8FlnyPv+yRJULsNsmfJd/W61u9 P/qKG5CrpVvwRFTYE3rseE5IwxPhDjhR3tiL04MOGWN+NjvN499e38bKxbOcbc339/2vVJPnvxuB eP9LiqZ8+dsusKL4HfK/Is9/N4KY//FLkeT/dmHx9rK+InA//6ur5z9l05T83wRS/F98CZJFYGuQ /mqxrgpwL/+r+ir/a7WS5P8msMp/cdQhK8CWIJx5+9Hleu9BKV6rVe7if7kan/9WzErJoPzXa7rk /0bw4IfiLAqL565XdLwPcG5Hlwo9EtacmQ+BGzgXtjtRFPcCzkDNPlBB8xww4e1jIJeOp2Sc4aUP 6iyyx04DsjoEfkgM9mmq8ORHEy0+ugQM5cJVFNZ5oGYNlTVNbJqqQkLbi+whcX0vQs316fPDVufo xcB61dDoidZcVUY2sXFqDu9+2rSah81+q6EFfkTGoROhSRD6xMfSlfZw2uta3aPucUNzPhLHGzkj NETz2YQMLvxwapO0da/Vf3lsDZ51eydNq6HRIxHqd8weiwOMUNr4sN1paPvFKByy2Amr+IrjZpET DuxwfJAvKJc4zbjNQ6l5wJy97Ld6DW2uLgK6GKe9RBtuohZY/JaGvuj2reWhi9tol2woNRFDo2Ho BiQ6iI/jk2e9OI5flutpefVfRJLD+JTmxp+RxTl8Sh4t/aeZOIdfkVe+iSwfwS/UGEWsWwMRbAzq tZKZ+EN7whKPJ5hQ8HXzRBMqf0Yu3AnNpZKqKBk1m9piFd4oGVRdx7E8+/ntPFHGe7NQagEGmt5U iB6/DCGVKkfH7VbH6mMyx4Pepbt/6R6m+06oQ5HLQqXxlNV4ymL/UgoLI6pPM0moL1DNYxCvIuYS 0lPNilioyhxD6tijwdQhuI/pkIpYLUIa2IQ4ocdjqmTsP9+D9gxyBznQPqQ6RUuFHN42a8BfcR+g 60xmHM3O88Xfz+ANefvoM79ki3ugqnuQNQvUJAomLslnzT0cGZII+0BlHUGIOcqVZ8ZbqqFFBq9z JZPDBSdLItNg5IY4mfz0PXGmAWijAis4AeTCKWghDQ83UnP0z1wWFinqHPVLZWk3gqQGYXs4cR2P MPU7/5xdk/qD7eUSQzt5hmHzDU+x1b1Kb8tdqXNXzqTz5UZuLGd3srhdzahHsEt/jXLyYfKPeI5x 3qg8wdUJMtgbfhpM6YRJEHHVILRxEVwj2oqCtxfUA5fVK1F/GHMew8hXMnE95z27lNhYNjOYkfQR wTelmL2mZvNB9po9O+b72M2NzLuMTGGkZFI1QrDUUFN0YNlP0/qmnXnDjqU6rpbOLZ9iSuwliQ7Y H5wQn4YqJloGQ3L3ABpBaoQDzVuNzLu93j1AeMVV0a2gD9Q8Y+gHELOnF1XIJpdNJovJ0kssm1zG /txh63m7w6ib7DkcQJ56gyegF+An5hCK7D7QAB1Nk5SgptRxbEqdoilTcdMkM/dLF0A/sA4kd9qD xJOSmedE1LiC5hBf6+4uPJzz+Kx2PXjwEOY0Kl9DgEw6B5L04fGL46Rq7PdfOjaFY1M4juWEW9kU uUa+58jvYRISEhISEhISEhISEhISEhIS/wh/AwhuQvcAUAAA --=-=-=-- ^ permalink raw reply [nested|flat] 278+ messages in thread
end of thread, other threads:[~2026-06-02 10:08 UTC | newest] Thread overview: 278+ messages (download: mbox mbox.gz follow: Atom feed) -- links below jump to the message on this page -- 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-06-02 10:08 [PATCH v2 4/4] Let pgbench support binary format. Andy Fan <[email protected]>
This inbox is served by agora; see mirroring instructions for how to clone and mirror all data and code used for this inbox