agora inbox for [email protected]help / color / mirror / Atom feed
[PATCH 2/5] Refactor index_concurrently_create_copy() for use with REPACK (CONCURRENTLY). 197+ messages / 2 participants [nested] [flat]
* [PATCH 2/5] Refactor index_concurrently_create_copy() for use with REPACK (CONCURRENTLY). @ 2026-03-05 18:58 Antonin Houska <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Antonin Houska @ 2026-03-05 18:58 UTC (permalink / raw) This patch moves the code to index_create_copy() and adds a "concurrently" parameter so it can be used by REPACK (CONCURRENTLY). With the CONCURRENTLY option, REPACK cannot simply swap the heap file and rebuild its indexes. Instead, it needs to build a separate set of indexes (including system catalog entries) *before* the actual swap, to reduce the time AccessExclusiveLock needs to be held for. --- src/backend/catalog/index.c | 54 +++++++++++++++++++++++--------- src/backend/commands/indexcmds.c | 6 ++-- src/backend/nodes/makefuncs.c | 9 +++--- src/include/catalog/index.h | 3 ++ src/include/nodes/makefuncs.h | 4 ++- 5 files changed, 54 insertions(+), 22 deletions(-) diff --git a/src/backend/catalog/index.c b/src/backend/catalog/index.c index 5ee6389d39c..f8e6c3d804e 100644 --- a/src/backend/catalog/index.c +++ b/src/backend/catalog/index.c @@ -1288,15 +1288,32 @@ index_create(Relation heapRelation, /* * index_concurrently_create_copy * - * Create concurrently an index based on the definition of the one provided by - * caller. The index is inserted into catalogs and needs to be built later - * on. This is called during concurrent reindex processing. - * - * "tablespaceOid" is the tablespace to use for this index. + * Variant of index_create_copy(), called during concurrent reindex + * processing. */ Oid index_concurrently_create_copy(Relation heapRelation, Oid oldIndexId, Oid tablespaceOid, const char *newName) +{ + return index_create_copy(heapRelation, oldIndexId, tablespaceOid, newName, + true); +} + +/* + * index_create_copy + * + * Create an index based on the definition of the one provided by caller. The + * index is inserted into catalogs. If 'concurrently' is TRUE, it needs to be + * built later on, otherwise it's built immediately. + * + * "tablespaceOid" is the tablespace to use for this index. + * + * The actual implementation of index_concurrently_create_copy(), reusable for + * other purposes. + */ +Oid +index_create_copy(Relation heapRelation, Oid oldIndexId, Oid tablespaceOid, + const char *newName, bool concurrently) { Relation indexRelation; IndexInfo *oldInfo, @@ -1315,6 +1332,7 @@ index_concurrently_create_copy(Relation heapRelation, Oid oldIndexId, List *indexColNames = NIL; List *indexExprs = NIL; List *indexPreds = NIL; + int flags = 0; indexRelation = index_open(oldIndexId, RowExclusiveLock); @@ -1325,7 +1343,7 @@ index_concurrently_create_copy(Relation heapRelation, Oid oldIndexId, * Concurrent build of an index with exclusion constraints is not * supported. */ - if (oldInfo->ii_ExclusionOps != NULL) + if (oldInfo->ii_ExclusionOps != NULL && concurrently) ereport(ERROR, (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), errmsg("concurrent index creation for exclusion constraints is not supported"))); @@ -1381,9 +1399,7 @@ index_concurrently_create_copy(Relation heapRelation, Oid oldIndexId, } /* - * Build the index information for the new index. Note that rebuild of - * indexes with exclusion constraints is not supported, hence there is no - * need to fill all the ii_Exclusion* fields. + * Build the index information for the new index. */ newInfo = makeIndexInfo(oldInfo->ii_NumIndexAttrs, oldInfo->ii_NumIndexKeyAttrs, @@ -1392,10 +1408,13 @@ index_concurrently_create_copy(Relation heapRelation, Oid oldIndexId, indexPreds, oldInfo->ii_Unique, oldInfo->ii_NullsNotDistinct, - false, /* not ready for inserts */ - true, + !concurrently, /* isready */ + concurrently, /* concurrent */ indexRelation->rd_indam->amsummarizing, - oldInfo->ii_WithoutOverlaps); + oldInfo->ii_WithoutOverlaps, + oldInfo->ii_ExclusionOps, + oldInfo->ii_ExclusionProcs, + oldInfo->ii_ExclusionStrats); /* * Extract the list of column names and the column numbers for the new @@ -1433,6 +1452,9 @@ index_concurrently_create_copy(Relation heapRelation, Oid oldIndexId, stattargets[i].isnull = isnull; } + if (concurrently) + flags = INDEX_CREATE_SKIP_BUILD | INDEX_CREATE_CONCURRENT; + /* * Now create the new index. * @@ -1456,7 +1478,7 @@ index_concurrently_create_copy(Relation heapRelation, Oid oldIndexId, indcoloptions->values, stattargets, reloptionsDatum, - INDEX_CREATE_SKIP_BUILD | INDEX_CREATE_CONCURRENT, + flags, 0, true, /* allow table to be a system catalog? */ false, /* is_internal? */ @@ -2450,7 +2472,8 @@ BuildIndexInfo(Relation index) indexStruct->indisready, false, index->rd_indam->amsummarizing, - indexStruct->indisexclusion && indexStruct->indisunique); + indexStruct->indisexclusion && indexStruct->indisunique, + NULL, NULL, NULL); /* fill in attribute numbers */ for (i = 0; i < numAtts; i++) @@ -2510,7 +2533,8 @@ BuildDummyIndexInfo(Relation index) indexStruct->indisready, false, index->rd_indam->amsummarizing, - indexStruct->indisexclusion && indexStruct->indisunique); + indexStruct->indisexclusion && indexStruct->indisunique, + NULL, NULL, NULL); /* fill in attribute numbers */ for (i = 0; i < numAtts; i++) diff --git a/src/backend/commands/indexcmds.c b/src/backend/commands/indexcmds.c index 635679cc1f2..34209bd1393 100644 --- a/src/backend/commands/indexcmds.c +++ b/src/backend/commands/indexcmds.c @@ -243,7 +243,8 @@ CheckIndexCompatible(Oid oldId, */ indexInfo = makeIndexInfo(numberOfAttributes, numberOfAttributes, accessMethodId, NIL, NIL, false, false, - false, false, amsummarizing, isWithoutOverlaps); + false, false, amsummarizing, isWithoutOverlaps, + NULL, NULL, NULL); typeIds = palloc_array(Oid, numberOfAttributes); collationIds = palloc_array(Oid, numberOfAttributes); opclassIds = palloc_array(Oid, numberOfAttributes); @@ -930,7 +931,8 @@ DefineIndex(ParseState *pstate, !concurrent, concurrent, amissummarizing, - stmt->iswithoutoverlaps); + stmt->iswithoutoverlaps, + NULL, NULL, NULL); typeIds = palloc_array(Oid, numberOfAttributes); collationIds = palloc_array(Oid, numberOfAttributes); diff --git a/src/backend/nodes/makefuncs.c b/src/backend/nodes/makefuncs.c index 2caec621d73..ca7e21e8349 100644 --- a/src/backend/nodes/makefuncs.c +++ b/src/backend/nodes/makefuncs.c @@ -834,7 +834,8 @@ IndexInfo * makeIndexInfo(int numattrs, int numkeyattrs, Oid amoid, List *expressions, List *predicates, bool unique, bool nulls_not_distinct, bool isready, bool concurrent, bool summarizing, - bool withoutoverlaps) + bool withoutoverlaps, Oid *exclusion_ops, Oid *exclusion_procs, + uint16 *exclusion_strats) { IndexInfo *n = makeNode(IndexInfo); @@ -863,9 +864,9 @@ makeIndexInfo(int numattrs, int numkeyattrs, Oid amoid, List *expressions, n->ii_PredicateState = NULL; /* exclusion constraints */ - n->ii_ExclusionOps = NULL; - n->ii_ExclusionProcs = NULL; - n->ii_ExclusionStrats = NULL; + n->ii_ExclusionOps = exclusion_ops; + n->ii_ExclusionProcs = exclusion_procs; + n->ii_ExclusionStrats = exclusion_strats; /* speculative inserts */ n->ii_UniqueOps = NULL; diff --git a/src/include/catalog/index.h b/src/include/catalog/index.h index b259c4141ed..3426087b445 100644 --- a/src/include/catalog/index.h +++ b/src/include/catalog/index.h @@ -99,6 +99,9 @@ extern Oid index_concurrently_create_copy(Relation heapRelation, Oid oldIndexId, Oid tablespaceOid, const char *newName); +extern Oid index_create_copy(Relation heapRelation, Oid oldIndexId, + Oid tablespaceOid, const char *newName, + bool concurrently); extern void index_concurrently_build(Oid heapRelationId, Oid indexRelationId); diff --git a/src/include/nodes/makefuncs.h b/src/include/nodes/makefuncs.h index 982ec25ae14..dcea148ae1a 100644 --- a/src/include/nodes/makefuncs.h +++ b/src/include/nodes/makefuncs.h @@ -99,7 +99,9 @@ extern IndexInfo *makeIndexInfo(int numattrs, int numkeyattrs, Oid amoid, List *expressions, List *predicates, bool unique, bool nulls_not_distinct, bool isready, bool concurrent, - bool summarizing, bool withoutoverlaps); + bool summarizing, bool withoutoverlaps, + Oid *exclusion_ops, Oid *exclusion_procs, + uint16 *exclusion_strats); extern Node *makeStringConst(char *str, int location); extern DefElem *makeDefElem(char *name, Node *arg, int location); -- 2.47.3 --=-=-= Content-Type: text/plain Content-Disposition: attachment; filename=v38-0003-Add-CONCURRENTLY-option-to-REPACK-command.patch ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <[email protected]> 0 siblings, 0 replies; 197+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <[email protected]> Discussion: https://postgr.es/m/[email protected] --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 197+ messages in thread
end of thread, other threads:[~2026-04-07 20:29 UTC | newest] Thread overview: 197+ messages (download: mbox mbox.gz follow: Atom feed) -- links below jump to the message on this page -- 2026-03-05 18:58 [PATCH 2/5] Refactor index_concurrently_create_copy() for use with REPACK (CONCURRENTLY). Antonin Houska <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <[email protected]>
This inbox is served by agora; see mirroring instructions for how to clone and mirror all data and code used for this inbox